.. code:: ipython3

    #S11
    
    import pandas as pd
    
    data = {
        "Name": ["Piyush", "Ketaki", "Khushi", "Rakesh", "Rinkesh", "Gayatri", "Prajakta", "Ishika"],
        "Department": ["IT", "HR", "Sales", "IT", "Finance", "HR", "Sales", "IT"],
        "Designation": ["Manager", "Clerk", "Executive", "Developer", "Accountant", "HR Executive", "Salesman", "Developer"],
        "Salary": [50000, 30000, 35000, 45000, 40000, 32000, 28000, 42000]
    }
    
    df = pd.DataFrame(data)
    
    print("First 5 Records:")
    print(df.head())


.. parsed-literal::

    First 5 Records:
          Name Department Designation  Salary
    0   Piyush         IT     Manager   50000
    1   Ketaki         HR       Clerk   30000
    2   Khushi      Sales   Executive   35000
    3   Rakesh         IT   Developer   45000
    4  Rinkesh    Finance  Accountant   40000
    

.. code:: ipython3

    #S13
    
    import numpy as np
    import matplotlib.pyplot as plt
    
    data = np.random.randint(1, 100, 25)
    
    # Line Chart
    plt.plot(data)
    plt.title("Line Chart")
    plt.xlabel("Index")
    plt.ylabel("Value")
    plt.show()
    
    # Scatter Plot
    plt.scatter(range(25), data)
    plt.title("Scatter Plot")
    plt.xlabel("Index")
    plt.ylabel("Value")
    plt.show()
    
    # Histogram
    plt.hist(data)
    plt.title("Histogram")
    plt.xlabel("Value")
    plt.ylabel("Frequency")
    plt.show()
    
    # Box Plot
    plt.boxplot(data)
    plt.title("Box Plot")
    plt.ylabel("Value")
    plt.show()



.. image:: output_1_0.png



.. image:: output_1_1.png



.. image:: output_1_2.png



.. image:: output_1_3.png


.. code:: ipython3

    #S21
    
    import pandas as pd
    
    marks = [32, 80, 71, 98, 82, 58, 94]
    
    data = pd.Series(marks)
    
    print("Sum =", data.sum())
    print("Average =", data.mean())
    print("Median =", data.median())
    print("Mode =", data.mode()[0])
    print("Minimum =", data.min())
    print("Maximum =", data.max())
    print("Standard Deviation =", data.std())


.. parsed-literal::

    Sum = 515
    Average = 73.57142857142857
    Median = 80.0
    Mode = 32
    Minimum = 32
    Maximum = 98
    Standard Deviation = 22.74391598396622
    

.. code:: ipython3

    import matplotlib.pyplot as plt
    
    names = ["Piyush","Ketaki", "Khushi", "Rakesh", "Rinkesh"]
    percentage = [32, 80, 71, 98, 82]
    
    # Pie Chart
    plt.pie(percentage, labels=names, autopct="%1.1f%%")
    plt.title("Student Percentage")
    plt.show()
    
    # Bar Chart
    plt.bar(names, percentage)
    plt.title("Student Percentage")
    plt.xlabel("Students")
    plt.ylabel("Percentage")
    plt.show()



.. image:: output_3_0.png



.. image:: output_3_1.png


.. code:: ipython3

    #S31
    
    import pandas as pd
    
    data = {
        "Name": ["Piyush","Ketaki", "Khushi", "Rakesh", "Rinkesh"],
        "Attendance": [85, 90, 78, 88, 85],
        "Score": [80, None, 75, None, 80]
    }
    
    df = pd.DataFrame(data)
    
    print("Original Data:")
    print(df)
    
    # Replace missing scores with mean
    mean_score = df["Score"].mean()
    df["Score"] = df["Score"].fillna(mean_score)
    
    print("\nAfter filling missing scores:")
    print(df)
    
    # Find duplicate records
    print("\nDuplicate Records:")
    print(df.duplicated())


.. parsed-literal::

    Original Data:
          Name  Attendance  Score
    0   Piyush          85   80.0
    1   Ketaki          90    NaN
    2   Khushi          78   75.0
    3   Rakesh          88    NaN
    4  Rinkesh          85   80.0
    
    After filling missing scores:
          Name  Attendance      Score
    0   Piyush          85  80.000000
    1   Ketaki          90  78.333333
    2   Khushi          78  75.000000
    3   Rakesh          88  78.333333
    4  Rinkesh          85  80.000000
    
    Duplicate Records:
    0    False
    1    False
    2    False
    3    False
    4    False
    dtype: bool
    

.. code:: ipython3

    #S33
    
    import numpy as np
    import matplotlib.pyplot as plt
    
    ages = np.array([22, 25, 28, 30, 32, 35, 27, 29, 31, 40, 42, 45, 38, 36, 50])
    
    bins = [20, 30, 40, 50, 60]
    
    frequency, edges = np.histogram(ages, bins)
    
    print("Frequency of each bin:")
    print(frequency)
    
    plt.hist(ages, bins=bins, color="blue", edgecolor="black")
    plt.title("Employee Age Histogram")
    plt.xlabel("Age")
    plt.ylabel("Frequency")
    plt.show()


.. parsed-literal::

    Frequency of each bin:
    [5 6 3 1]
    


.. image:: output_5_1.png


.. code:: ipython3

    #S41
    
    import pandas as pd
    
    data = {
        "Product": ["Laptop", "Mobile", "Tablet", "Mouse", "Keyboard", "Monitor", "Printer", "Speaker"],
        "Brand": ["Dell", "Samsung", "Apple", "HP", "Logitech", "LG", "Canon", "Boat"],
        "Category": ["Computer", "Phone", "Electronics", "Accessory", "Accessory", "Computer", "Printer", "Audio"],
        "Price": [50000, 25000, 40000, 800, 1200, 15000, 10000, 2000]
    }
    
    df = pd.DataFrame(data)
    
    print(df.head())


.. parsed-literal::

        Product     Brand     Category  Price
    0    Laptop      Dell     Computer  50000
    1    Mobile   Samsung        Phone  25000
    2    Tablet     Apple  Electronics  40000
    3     Mouse        HP    Accessory    800
    4  Keyboard  Logitech    Accessory   1200
    

.. code:: ipython3

    #S42
    
    import pandas as pd
    import numpy as np
    
    salary = [25, 30, 35, 40, 45, 50, 55, 60, 65, 70, 75, 80, 85, 90, 150]
    
    df = pd.DataFrame({"Salary": salary})
    
    # Quartiles
    Q1 = df["Salary"].quantile(0.25)
    Q2 = df["Salary"].quantile(0.50)
    Q3 = df["Salary"].quantile(0.75)
    
    print("Q1 =", Q1)
    print("Q2 =", Q2)
    print("Q3 =", Q3)
    
    # Percentiles
    print("25th Percentile =", df["Salary"].quantile(0.25))
    print("50th Percentile =", df["Salary"].quantile(0.50))
    print("75th Percentile =", df["Salary"].quantile(0.75))
    print("90th Percentile =", df["Salary"].quantile(0.90))
    
    # IQR Method
    IQR = Q3 - Q1
    lower = Q1 - 1.5 * IQR
    upper = Q3 + 1.5 * IQR
    
    outliers = df[(df["Salary"] < lower) | (df["Salary"] > upper)]
    
    print("IQR Outliers:")
    print(outliers)
    
    # Z-score Method
    mean = df["Salary"].mean()
    std = df["Salary"].std()
    
    df["Z-score"] = (df["Salary"] - mean) / std
    
    print("Z-score Outliers:")
    print(df[abs(df["Z-score"]) > 3])


.. parsed-literal::

    Q1 = 42.5
    Q2 = 60.0
    Q3 = 77.5
    25th Percentile = 42.5
    50th Percentile = 60.0
    75th Percentile = 77.5
    90th Percentile = 88.0
    IQR Outliers:
        Salary
    14     150
    Z-score Outliers:
    Empty DataFrame
    Columns: [Salary, Z-score]
    Index: []
    

.. code:: ipython3

    #S51
    
    import pandas as pd
    
    data = {
        "Laptop_Model": ["Dell", "HP", "Lenovo", "Acer", "Samsung"],
        "Cost": [50000, 55000, 48000, 45000, 62000]
    }
    
    df = pd.DataFrame(data)
    
    print(df)


.. parsed-literal::

      Laptop_Model   Cost
    0         Dell  50000
    1           HP  55000
    2       Lenovo  48000
    3         Acer  45000
    4      Samsung  62000
    

.. code:: ipython3

    #S52
    
    import matplotlib.pyplot as plt
    
    names = ["Piyush","Ketaki", "Khushi", "Rakesh", "Rinkesh"]
    percentage = [85, 78, 92, 75, 88]
    
    # Pie Chart
    plt.pie(percentage, labels=names, autopct="%1.1f%%")
    plt.title("Student Percentage")
    plt.show()
    
    # Bar Chart
    plt.bar(names, percentage)
    plt.title("Student Percentage")
    plt.xlabel("Student Name")
    plt.ylabel("Percentage")
    plt.show()



.. image:: output_9_0.png



.. image:: output_9_1.png


.. code:: ipython3

    #S61
    
    import pandas as pd
    
    data = {
        "Name": ["Piyush", "Ketaki", "Khushi", "Rakesh", "Rinkesh", "Gayatri", "Prajakta", "Ishika"],
        "Age": [20, 21, 19, 22, 20, 21, 20, 22],
        "Course": ["BCA", "BCA", "BSc", "BCom",
                   "BCA", "BBA", "BCA", "BSc"],
        "Marks": [85, 78, 90, 75, 88, 82, 91, 79]
    }
    
    df = pd.DataFrame(data)
    
    print(df.head())


.. parsed-literal::

          Name  Age Course  Marks
    0   Piyush   20    BCA     85
    1   Ketaki   21    BCA     78
    2   Khushi   19    BSc     90
    3   Rakesh   22   BCom     75
    4  Rinkesh   20    BCA     88
    

.. code:: ipython3

    #S63
    
    import pandas as pd
    
    data = {
        "Name": ["Piyush", "Ketaki", "Khushi", "Rakesh", "Rinkesh"],
        "Address": ["Pune", "Pune", "Nashik", "Chennai", "Madurai"],
        "Age": [20, 23, 19, 21, 22],
        "Class": ["ME", "MSc (CA)", "MVoC", "MCom", "MSc"]
    }
    
    df = pd.DataFrame(data)
    
    # Save as CSV
    df.to_csv("StudentData.csv", index=False)
    
    # Import Dataset
    df = pd.read_csv("StudentData.csv")
    
    # a) Statistical description
    print("Statistical Description:")
    print(df.describe())
    
    # b) Shape
    print("\nShape:")
    print(df.shape)
    
    # c) First two rows
    print("\nFirst Two Rows:")
    print(df.head(2))


.. parsed-literal::

    Statistical Description:
                 Age
    count   5.000000
    mean   21.000000
    std     1.581139
    min    19.000000
    25%    20.000000
    50%    21.000000
    75%    22.000000
    max    23.000000
    
    Shape:
    (5, 4)
    
    First Two Rows:
         Name Address  Age     Class
    0  Piyush    Pune   20        ME
    1  Ketaki    Pune   23  MSc (CA)
    

.. code:: ipython3

    #S71
    
    import pandas as pd
    
    data = {
        "staff_id": [101, 102, 103, 104, 105],
        "employee_name": ["Piyush", "Ketaki", "Khushi", "Rakesh", "Rinkesh"],
        "department": ["IT", "HR", "Sales", "Finance", "IT"],
        "monthly_salary": [30000, 35000, 32000, 40000, 38000]
    }
    
    df = pd.DataFrame(data)
    
    # Create CSV file
    df.to_csv("staff.csv", index=False)
    
    # Read CSV file
    df = pd.read_csv("staff.csv")
    
    print(df)


.. parsed-literal::

       staff_id employee_name department  monthly_salary
    0       101        Piyush         IT           30000
    1       102        Ketaki         HR           35000
    2       103        Khushi      Sales           32000
    3       104        Rakesh    Finance           40000
    4       105       Rinkesh         IT           38000
    

.. code:: ipython3

    #S72
    
    
    import numpy as np
    import matplotlib.pyplot as plt
    
    ages = np.array([22, 25, 28, 30, 32, 35, 27, 29, 31, 40, 42, 45, 38, 36, 50])
    
    bins = [20, 30, 40, 50, 60]
    
    # Calculate frequency
    frequency, edges = np.histogram(ages, bins)
    
    print("Frequency:", frequency)
    
    # Plot histogram
    plt.hist(ages, bins=bins, color="blue", edgecolor="black")
    
    plt.title("Employee Age Histogram")
    plt.xlabel("Age")
    plt.ylabel("Frequency")
    
    plt.show()


.. parsed-literal::

    Frequency: [5 6 3 1]
    


.. image:: output_13_1.png


.. code:: ipython3

    #S81
    
    import pandas as pd
    
    data = {
        "Book_Title": ["Harry Potter", "Wings of Fire", "The Alchemist",
                       "Atomic Habits", "Rich Dad Poor Dad",
                       "The Hobbit", "Ikigai", "The Monk Who Sold His Ferrari"],
        "Author": ["J.K. Rowling", "A.P.J. Abdul Kalam", "Paulo Coelho",
                   "James Clear", "Robert Kiyosaki", "J.R.R. Tolkien",
                   "Hector Garcia", "Robin Sharma"],
        "Category": ["Fantasy", "Biography", "Fiction", "Self-help",
                     "Finance", "Fantasy", "Self-help", "Self-help"],
        "Rating": [4.8, 4.7, 4.6, 4.8, 4.5, 4.7, 4.4, 4.6]
    }
    
    df = pd.DataFrame(data)
    
    print(df.head())


.. parsed-literal::

              Book_Title              Author   Category  Rating
    0       Harry Potter        J.K. Rowling    Fantasy     4.8
    1      Wings of Fire  A.P.J. Abdul Kalam  Biography     4.7
    2      The Alchemist        Paulo Coelho    Fiction     4.6
    3      Atomic Habits         James Clear  Self-help     4.8
    4  Rich Dad Poor Dad     Robert Kiyosaki    Finance     4.5
    

.. code:: ipython3

    #S83
    
    import pandas as pd
    
    salary = [25, 30, 35, 40, 45, 50, 55, 60, 65, 70, 75, 80, 85, 90, 150]
    
    df = pd.DataFrame({"Salary": salary})
    
    # Quartiles
    Q1 = df["Salary"].quantile(0.25)
    Q2 = df["Salary"].quantile(0.50)
    Q3 = df["Salary"].quantile(0.75)
    
    print("Q1 =", Q1)
    print("Q2 =", Q2)
    print("Q3 =", Q3)
    
    # Percentiles
    print("25th =", df["Salary"].quantile(0.25))
    print("50th =", df["Salary"].quantile(0.50))
    print("75th =", df["Salary"].quantile(0.75))
    print("90th =", df["Salary"].quantile(0.90))
    
    # IQR
    IQR = Q3 - Q1
    lower = Q1 - 1.5 * IQR
    upper = Q3 + 1.5 * IQR
    
    print("IQR Outliers:")
    print(df[(df["Salary"] < lower) | (df["Salary"] > upper)])
    
    # Z-score
    mean = df["Salary"].mean()
    std = df["Salary"].std()
    
    df["Z"] = (df["Salary"] - mean) / std
    
    print("Z-score Outliers:")
    print(df[abs(df["Z"]) > 3])


.. parsed-literal::

    Q1 = 42.5
    Q2 = 60.0
    Q3 = 77.5
    25th = 42.5
    50th = 60.0
    75th = 77.5
    90th = 88.0
    IQR Outliers:
        Salary
    14     150
    Z-score Outliers:
    Empty DataFrame
    Columns: [Salary, Z]
    Index: []
    

.. code:: ipython3

    #S91
    
    import pandas as pd
    
    df = pd.read_csv("iris.csv")
    
    frequency = df["Species"].value_counts()
    
    print("Frequency of Iris Species:")
    print(frequency)
    
    # we have to import iris dataset 


::


    ---------------------------------------------------------------------------

    KeyError                                  Traceback (most recent call last)

    File D:\Anaconda\Lib\site-packages\pandas\core\indexes\base.py:3812, in Index.get_loc(self, key)
       3811 try:
    -> 3812     return self._engine.get_loc(casted_key)
       3813 except KeyError as err:
    

    File pandas/_libs/index.pyx:167, in pandas._libs.index.IndexEngine.get_loc()
    

    File pandas/_libs/index.pyx:196, in pandas._libs.index.IndexEngine.get_loc()
    

    File pandas/_libs/hashtable_class_helper.pxi:7088, in pandas._libs.hashtable.PyObjectHashTable.get_item()
    

    File pandas/_libs/hashtable_class_helper.pxi:7096, in pandas._libs.hashtable.PyObjectHashTable.get_item()
    

    KeyError: 'Species'

    
    The above exception was the direct cause of the following exception:
    

    KeyError                                  Traceback (most recent call last)

    Cell In[13], line 7
          3 import pandas as pd
          5 df = pd.read_csv("iris.csv")
    ----> 7 frequency = df["Species"].value_counts()
          9 print("Frequency of Iris Species:")
         10 print(frequency)
    

    File D:\Anaconda\Lib\site-packages\pandas\core\frame.py:4113, in DataFrame.__getitem__(self, key)
       4111 if self.columns.nlevels > 1:
       4112     return self._getitem_multilevel(key)
    -> 4113 indexer = self.columns.get_loc(key)
       4114 if is_integer(indexer):
       4115     indexer = [indexer]
    

    File D:\Anaconda\Lib\site-packages\pandas\core\indexes\base.py:3819, in Index.get_loc(self, key)
       3814     if isinstance(casted_key, slice) or (
       3815         isinstance(casted_key, abc.Iterable)
       3816         and any(isinstance(x, slice) for x in casted_key)
       3817     ):
       3818         raise InvalidIndexError(key)
    -> 3819     raise KeyError(key) from err
       3820 except TypeError:
       3821     # If we have a listlike key, _check_indexing_error will raise
       3822     #  InvalidIndexError. Otherwise we fall through and re-raise
       3823     #  the TypeError.
       3824     self._check_indexing_error(key)
    

    KeyError: 'Species'


.. code:: ipython3

    #S93
    
    import pandas as pd
    
    df = pd.read_csv("employee.csv")
    
    salary = df["Salary"]
    
    print("Count =", salary.count())
    print("Mean =", salary.mean())
    print("Median =", salary.median())
    print("Variance =", salary.var())
    print("Standard Deviation =", salary.std())
    print("Minimum =", salary.min())
    print("Maximum =", salary.max())
    
    print("Q1 =", salary.quantile(0.25))
    print("Q2 =", salary.quantile(0.50))
    print("Q3 =", salary.quantile(0.75))
    
    # we have to import iris dataset 


::


    ---------------------------------------------------------------------------

    FileNotFoundError                         Traceback (most recent call last)

    Cell In[1], line 5
          1 #S93
          3 import pandas as pd
    ----> 5 df = pd.read_csv("employee.csv")
          7 salary = df["Salary"]
          9 print("Count =", salary.count())
    

    File D:\Anaconda\Lib\site-packages\pandas\io\parsers\readers.py:1026, in read_csv(filepath_or_buffer, sep, delimiter, header, names, index_col, usecols, dtype, engine, converters, true_values, false_values, skipinitialspace, skiprows, skipfooter, nrows, na_values, keep_default_na, na_filter, verbose, skip_blank_lines, parse_dates, infer_datetime_format, keep_date_col, date_parser, date_format, dayfirst, cache_dates, iterator, chunksize, compression, thousands, decimal, lineterminator, quotechar, quoting, doublequote, escapechar, comment, encoding, encoding_errors, dialect, on_bad_lines, delim_whitespace, low_memory, memory_map, float_precision, storage_options, dtype_backend)
       1013 kwds_defaults = _refine_defaults_read(
       1014     dialect,
       1015     delimiter,
       (...)   1022     dtype_backend=dtype_backend,
       1023 )
       1024 kwds.update(kwds_defaults)
    -> 1026 return _read(filepath_or_buffer, kwds)
    

    File D:\Anaconda\Lib\site-packages\pandas\io\parsers\readers.py:620, in _read(filepath_or_buffer, kwds)
        617 _validate_names(kwds.get("names", None))
        619 # Create the parser.
    --> 620 parser = TextFileReader(filepath_or_buffer, **kwds)
        622 if chunksize or iterator:
        623     return parser
    

    File D:\Anaconda\Lib\site-packages\pandas\io\parsers\readers.py:1620, in TextFileReader.__init__(self, f, engine, **kwds)
       1617     self.options["has_index_names"] = kwds["has_index_names"]
       1619 self.handles: IOHandles | None = None
    -> 1620 self._engine = self._make_engine(f, self.engine)
    

    File D:\Anaconda\Lib\site-packages\pandas\io\parsers\readers.py:1880, in TextFileReader._make_engine(self, f, engine)
       1878     if "b" not in mode:
       1879         mode += "b"
    -> 1880 self.handles = get_handle(
       1881     f,
       1882     mode,
       1883     encoding=self.options.get("encoding", None),
       1884     compression=self.options.get("compression", None),
       1885     memory_map=self.options.get("memory_map", False),
       1886     is_text=is_text,
       1887     errors=self.options.get("encoding_errors", "strict"),
       1888     storage_options=self.options.get("storage_options", None),
       1889 )
       1890 assert self.handles is not None
       1891 f = self.handles.handle
    

    File D:\Anaconda\Lib\site-packages\pandas\io\common.py:873, in get_handle(path_or_buf, mode, encoding, compression, memory_map, is_text, errors, storage_options)
        868 elif isinstance(handle, str):
        869     # Check whether the filename is to be opened in binary mode.
        870     # Binary mode does not support 'encoding' and 'newline'.
        871     if ioargs.encoding and "b" not in ioargs.mode:
        872         # Encoding
    --> 873         handle = open(
        874             handle,
        875             ioargs.mode,
        876             encoding=ioargs.encoding,
        877             errors=errors,
        878             newline="",
        879         )
        880     else:
        881         # Binary mode
        882         handle = open(handle, ioargs.mode)
    

    FileNotFoundError: [Errno 2] No such file or directory: 'employee.csv'


.. code:: ipython3

    #S101
    
    import pandas as pd
    
    data = {
        "student_id": [1, 2, 3, 4, 5],
        "name": ["Piyush", "Ketaki", "Khushi", "Rakesh", "Rinkesh"],
        "marks": [75, 80, 85, 80, 90]
    }
    
    df = pd.DataFrame(data)
    
    print(df)
    
    print("Mean =", df["marks"].mean())
    print("Median =", df["marks"].median())
    print("Mode =", df["marks"].mode()[0])


.. parsed-literal::

       student_id     name  marks
    0           1   Piyush     75
    1           2   Ketaki     80
    2           3   Khushi     85
    3           4   Rakesh     80
    4           5  Rinkesh     90
    Mean = 82.0
    Median = 80.0
    Mode = 80
    

.. code:: ipython3

    #S103
    
    import pandas as pd
    
    df = pd.read_csv("Student_Marks.csv")
    
    print("Before Discretization:")
    print(df)
    
    # Convert marks into categories
    bins = [0, 40, 60, 80, 100]
    labels = ["Poor", "Average", "Good", "Excellent"]
    
    df["Performance"] = pd.cut(df["Marks"], bins=bins, labels=labels)
    
    print("\nAfter Discretization:")
    print(df)
    
    ## we have to import iris dataset 

.. code:: ipython3

    #S111
    
    import pandas as pd
    
    sales = [20, 25, 30, 28, 35, 40, 32, 38, 45, 42, 48, 50]
    
    df = pd.DataFrame({"Sales": sales})
    
    print(df)
    
    # Range
    print("Range =", df["Sales"].max() - df["Sales"].min())
    
    # Variance
    print("Variance =", df["Sales"].var())
    
    # Standard Deviation
    print("Standard Deviation =", df["Sales"].std())


.. parsed-literal::

        Sales
    0      20
    1      25
    2      30
    3      28
    4      35
    5      40
    6      32
    7      38
    8      45
    9      42
    10     48
    11     50
    Range = 30
    Variance = 88.26515151515152
    Standard Deviation = 9.394953513198004
    

.. code:: ipython3

    #S113
    
    import pandas as pd
    import matplotlib.pyplot as plt
    
    df = pd.read_csv("iris.csv")
    
    df.boxplot(column=["SepalLengthCm", "SepalWidthCm",
                       "PetalLengthCm", "PetalWidthCm"],
               by="Species")
    
    plt.suptitle("")
    plt.title("Iris Features Box Plot")
    plt.xlabel("Species")
    plt.ylabel("Values")
    plt.show()
    
    ## we have to import iris dataset 

.. code:: ipython3

    #S121
    
    import pandas as pd
    import matplotlib.pyplot as plt
    
    df = pd.read_csv("iris.csv")
    
    df.boxplot(
        column=["SepalLengthCm", "SepalWidthCm",
                "PetalLengthCm", "PetalWidthCm"],
        by="Species"
    )
    
    plt.suptitle("")
    plt.show()
    
    ## we have to import iris dataset 

.. code:: ipython3

    #S123
    
    import numpy as np
    import matplotlib.pyplot as plt
    
    data = np.random.randint(1, 100, 50)
    
    # Add two outliers
    data = np.append(data, [200, 250])
    
    # Line Chart
    plt.plot(data, color="blue")
    plt.title("Line Chart")
    plt.xlabel("Index")
    plt.ylabel("Value")
    plt.grid()
    plt.show()
    
    # Scatter Plot
    plt.scatter(range(len(data)), data, color="red")
    plt.title("Scatter Plot")
    plt.xlabel("Index")
    plt.ylabel("Value")
    plt.show()
    
    # Histogram
    plt.hist(data, color="green", edgecolor="black")
    plt.title("Histogram")
    plt.xlabel("Value")
    plt.ylabel("Frequency")
    plt.show()
    
    # Box Plot
    plt.boxplot(data)
    plt.title("Box Plot with Outliers")
    plt.ylabel("Value")
    plt.show()



.. image:: output_23_0.png



.. image:: output_23_1.png



.. image:: output_23_2.png



.. image:: output_23_3.png


.. code:: ipython3

    #S131
    
    import pandas as pd
    
    df = pd.read_csv("data.csv")
    
    print("Shape:", df.shape)
    print("Size:", df.size)
    print("\nData Types:")
    print(df.dtypes)
    
    print("\nStatistical Details:")
    print(df.describe())
    
    print("\nNumber of Observations:", len(df))
    
    print("\nMissing Values:")
    print(df.isnull().sum())

.. code:: ipython3

    #S133
    
    import numpy as np
    import matplotlib.pyplot as plt
    
    data = np.random.randint(1, 100, 50)
    
    # Line Chart
    plt.plot(data, color="blue")
    plt.title("Line Chart")
    plt.xlabel("Index")
    plt.ylabel("Value")
    plt.grid()
    plt.show()
    
    # Scatter Plot
    plt.scatter(range(50), data, color="red")
    plt.title("Scatter Plot")
    plt.xlabel("Index")
    plt.ylabel("Value")
    plt.show()
    
    # Histogram
    plt.hist(data, color="green", edgecolor="black")
    plt.title("Histogram")
    plt.xlabel("Value")
    plt.ylabel("Frequency")
    plt.show()
    
    # Box Plot
    plt.boxplot(data)
    plt.title("Box Plot")
    plt.ylabel("Value")
    plt.show()
    
    # Add two outliers
    data = np.append(data, [200, 250])
    
    # Updated Box Plot
    plt.boxplot(data)
    plt.title("Box Plot with Outliers")
    plt.ylabel("Value")
    plt.show()



.. image:: output_25_0.png



.. image:: output_25_1.png



.. image:: output_25_2.png



.. image:: output_25_3.png



.. image:: output_25_4.png


.. code:: ipython3

    #S141
    
    import pandas as pd
    
    data = {
        "Name": ["Amit", "Rahul", "Rohan", "Suresh", "Akash"],
        "Marks": [80, 75, 90, 85, 70],
        "City": ["Pune", "Mumbai", "Delhi", "Nashik", "Pune"]
    }
    
    df = pd.DataFrame(data)
    
    print(df)


.. parsed-literal::

         Name  Marks    City
    0    Amit     80    Pune
    1   Rahul     75  Mumbai
    2   Rohan     90   Delhi
    3  Suresh     85  Nashik
    4   Akash     70    Pune
    

.. code:: ipython3

    #S142
    import pandas as pd
    
    df = pd.read_csv("employee.csv")
    
    salary = df["Salary"]
    
    print("Count =", salary.count())
    print("Mean =", salary.mean())
    print("Median =", salary.median())
    print("Variance =", salary.var())
    print("Standard Deviation =", salary.std())
    print("Minimum =", salary.min())
    print("Maximum =", salary.max())
    
    print("Q1 =", salary.quantile(0.25))
    print("Q2 =", salary.quantile(0.50))
    print("Q3 =", salary.quantile(0.75))

.. code:: ipython3

    #S151
    
    import matplotlib.pyplot as plt
    
    subjects = ["Maths", "Python", "DBMS", "DS", "AI"]
    marks = [80, 75, 90, 85, 70]
    
    # Pie Chart
    plt.pie(marks, labels=subjects, autopct="%1.1f%%")
    plt.title("Subject Marks")
    plt.show()
    
    # Bar Chart
    plt.bar(subjects, marks, color="blue")
    plt.title("Subject Marks")
    plt.xlabel("Subjects")
    plt.ylabel("Marks")
    plt.show()



.. image:: output_28_0.png



.. image:: output_28_1.png


.. code:: ipython3

    #S152
    
    import pandas as pd
    
    df = pd.read_csv("data.csv")
    
    # Display DataFrame
    print("DataFrame:")
    print(df)
    
    # Describe
    print("\nDescribe:")
    print(df.describe())
    
    # First 5 rows
    print("\nHead:")
    print(df.head())
    
    # Last 5 rows
    print("\nTail:")
    print(df.tail())
    
    # Maximum
    print("\nMaximum:")
    print(df.max(numeric_only=True))
    
    # Minimum
    print("\nMinimum:")
    print(df.min(numeric_only=True))

.. code:: ipython3

    #S161
    
    import pandas as pd
    
    data = {
        "RollNo": [1, 2, 3, 4, 5],
        "Name": ["Amit", "Rahul", "Sneha", "Priya", "Rohan"],
        "Mark1": [70, 80, 65, 90, 75],
        "Mark2": [75, 85, 70, 95, 80],
        "Mark3": [80, 90, 75, 85, 70]
    }
    
    df = pd.DataFrame(data)
    
    print("DataFrame:")
    print(df)
    
    print("\nMaximum:")
    print(df.max(numeric_only=True))
    
    print("\nMinimum:")
    print(df.min(numeric_only=True))
    
    print("\nDescribe:")
    print(df.describe())
    
    print("\nShape:")
    print(df.shape)


.. parsed-literal::

    DataFrame:
       RollNo   Name  Mark1  Mark2  Mark3
    0       1   Amit     70     75     80
    1       2  Rahul     80     85     90
    2       3  Sneha     65     70     75
    3       4  Priya     90     95     85
    4       5  Rohan     75     80     70
    
    Maximum:
    RollNo     5
    Mark1     90
    Mark2     95
    Mark3     90
    dtype: int64
    
    Minimum:
    RollNo     1
    Mark1     65
    Mark2     70
    Mark3     70
    dtype: int64
    
    Describe:
             RollNo      Mark1      Mark2      Mark3
    count  5.000000   5.000000   5.000000   5.000000
    mean   3.000000  76.000000  81.000000  80.000000
    std    1.581139   9.617692   9.617692   7.905694
    min    1.000000  65.000000  70.000000  70.000000
    25%    2.000000  70.000000  75.000000  75.000000
    50%    3.000000  75.000000  80.000000  80.000000
    75%    4.000000  80.000000  85.000000  85.000000
    max    5.000000  90.000000  95.000000  90.000000
    
    Shape:
    (5, 5)
    

.. code:: ipython3

    #S162
    
    import numpy as np
    
    data = np.array([
        [10, 20, 30],
        [20, 30, 40],
        [30, 40, 50]
    ])
    
    weights = np.array([1, 2, 1])
    
    # Weighted Arithmetic Mean
    arithmetic = np.average(data, axis=0, weights=weights)
    
    # Weighted Harmonic Mean
    harmonic = np.sum(weights) / np.sum(weights / data, axis=0)
    
    # Weighted Geometric Mean
    geometric = np.exp(
        np.sum(weights * np.log(data), axis=0) / np.sum(weights)
    )
    
    print("Weighted Arithmetic Mean:", arithmetic)
    print("Weighted Harmonic Mean:", harmonic)
    print("Weighted Geometric Mean:", geometric)


.. parsed-literal::

    Weighted Arithmetic Mean: [20. 30. 40.]
    Weighted Harmonic Mean: [21.81818182 18.46153846 51.06382979]
    Weighted Geometric Mean: [  8.80111737 154.91933385  15.6508458 ]
    

.. code:: ipython3

    #S171
    
    import numpy as np
    
    ages = np.array([22, 25, 30, 28, 35, 40, 27, 32, 29, 45])
    
    print("Minimum Age:", np.min(ages))
    print("Maximum Age:", np.max(ages))
    print("Average Age:", np.mean(ages))


.. parsed-literal::

    Minimum Age: 22
    Maximum Age: 45
    Average Age: 31.3
    

.. code:: ipython3

    #S172
    
    import pandas as pd
    
    df = pd.read_csv("iris.csv")
    
    print("Dataset:")
    print(df)
    
    print("\nDescription:")
    print(df.describe())
    
    print("\nMean of Numeric Columns:")
    print(df.mean(numeric_only=True))
    
    print("\nMissing Values:")
    print(df.isnull().sum())

.. code:: ipython3

    #S181
    
    import numpy as np
    
    data = np.array([10, 20, 30, 40, 50, 60, 70, 80, 90, 100])
    
    print("Sum:", np.sum(data))
    print("Mean:", np.mean(data))
    print("Minimum:", np.min(data))
    print("Maximum:", np.max(data))


.. parsed-literal::

    Sum: 550
    Mean: 55.0
    Minimum: 10
    Maximum: 100
    

.. code:: ipython3

    #S182
    
    import pandas as pd
    
    # Read Iris dataset
    df = pd.read_csv("iris.csv")
    
    print("Sample Records:")
    print(df.sample(5))
    
    # Maximum values of numerical columns
    print("\nMaximum Values:")
    print(df.max(numeric_only=True))
    
    # Minimum values of numerical columns
    print("\nMinimum Values:")
    print(df.min(numeric_only=True))
    
    # Number of records for each class
    print("\nClass Counts:")
    print(df["Species"].value_counts())

.. code:: ipython3

    #S191
    
    import numpy as np
    
    marks = np.array([75, 88, 65, 92, 81, 70, 95, 78, 85, 90])
    
    marks = np.sort(marks)
    
    print("Sorted Marks:", marks)
    print("Highest 3 Marks:", marks[-3:])


.. parsed-literal::

    Sorted Marks: [65 70 75 78 81 85 88 90 92 95]
    Highest 3 Marks: [90 92 95]
    

.. code:: ipython3

    #s192
    
    import numpy as np
    
    data = np.array([
        [10, 20, 30],
        [20, 30, 40],
        [30, 40, 50]
    ])
    
    weights = np.array([1, 2, 1])
    
    # Weighted Arithmetic Mean
    arithmetic = np.average(data, axis=0, weights=weights)
    
    # Weighted Harmonic Mean
    harmonic = np.sum(weights) / np.sum(weights / data, axis=0)
    
    # Weighted Geometric Mean
    geometric = np.exp(
        np.sum(weights * np.log(data), axis=0) / np.sum(weights)
    )
    
    print("Weighted Arithmetic Mean:", arithmetic)
    print("Weighted Harmonic Mean:", harmonic)
    print("Weighted Geometric Mean:", geometric)


.. parsed-literal::

    Weighted Arithmetic Mean: [20. 30. 40.]
    Weighted Harmonic Mean: [21.81818182 18.46153846 51.06382979]
    Weighted Geometric Mean: [  8.80111737 154.91933385  15.6508458 ]
    

.. code:: ipython3

    #S201
    
    import numpy as np
    
    data = np.array([10, 20, 30, 40, 50, 60, 70, 80, 90, 100])
    
    variance = np.var(data)
    std = np.std(data)
    
    print("Variance:", variance)
    print("Standard Deviation:", std)


.. parsed-literal::

    Variance: 825.0
    Standard Deviation: 28.722813232690143
    

.. code:: ipython3

    #S203
    
    import pandas as pd
    
    data = {
        "Salary": [
            25000, 28000, 30000, 32000, 35000,
            36000, 38000, 40000, 42000, 45000,
            46000, 48000, 50000, 52000, 55000,
            58000, 60000, 65000, 70000, 150000
        ]
    }
    
    df = pd.DataFrame(data)
    
    print("Q1:", df["Salary"].quantile(0.25))
    print("Q2:", df["Salary"].quantile(0.50))
    print("Q3:", df["Salary"].quantile(0.75))
    
    print("25th Percentile:", df["Salary"].quantile(0.25))
    print("50th Percentile:", df["Salary"].quantile(0.50))
    print("75th Percentile:", df["Salary"].quantile(0.75))
    print("90th Percentile:", df["Salary"].quantile(0.90))
    
    # IQR
    Q1 = df["Salary"].quantile(0.25)
    Q3 = df["Salary"].quantile(0.75)
    
    IQR = Q3 - Q1
    
    lower = Q1 - 1.5 * IQR
    upper = Q3 + 1.5 * IQR
    
    outliers = df[(df["Salary"] < lower) | (df["Salary"] > upper)]
    
    print("\nIQR:", IQR)
    print("Outliers:")
    print(outliers)


.. parsed-literal::

    Q1: 35750.0
    Q2: 45500.0
    Q3: 55750.0
    25th Percentile: 35750.0
    50th Percentile: 45500.0
    75th Percentile: 55750.0
    90th Percentile: 65500.00000000001
    
    IQR: 20000.0
    Outliers:
        Salary
    19  150000
    

.. code:: ipython3

    #S211
    
    import numpy as np
    
    marks = np.array([35, 45, 55, 65, 72, 80, 90, 25, 48, 95])
    
    fail = 0
    average = 0
    excellent = 0
    
    for mark in marks:
        if mark < 40:
            fail += 1
        elif mark < 75:
            average += 1
        else:
            excellent += 1
    
    print("Fail Students:", fail)
    print("Average Students:", average)
    print("Excellent Students:", excellent)


.. parsed-literal::

    Fail Students: 2
    Average Students: 5
    Excellent Students: 3
    

.. code:: ipython3

    #S213
    
    import numpy as np
    import matplotlib.pyplot as plt
    
    data1 = np.random.normal(50, 10, 1000)
    data2 = np.random.normal(70, 15, 1000)
    
    plt.hist(data1, bins=20, alpha=0.5, edgecolor="black", label="Dataset 1")
    plt.hist(data2, bins=20, alpha=0.5, edgecolor="black", label="Dataset 2")
    
    plt.title("Two Normal Distributions")
    plt.xlabel("Values")
    plt.ylabel("Frequency")
    plt.legend()
    plt.show()



.. image:: output_41_0.png


.. code:: ipython3

    #S221
    
    import pandas as pd
    
    subjects = ["Maths", "Python", "DBMS", "DS", "AI"]
    marks = [80, 75, 90, 85, 70]
    
    df = pd.DataFrame({
        "Subject": subjects,
        "Marks": marks
    })
    
    print(df)


.. parsed-literal::

      Subject  Marks
    0   Maths     80
    1  Python     75
    2    DBMS     90
    3      DS     85
    4      AI     70
    

.. code:: ipython3

    #S223
    
    import matplotlib.pyplot as plt
    
    data = [20, 30, 25, 40, 35, 50]
    
    fig, ax = plt.subplots(figsize=(8, 5))
    
    ax.plot(data, linewidth=2)
    
    ax.set_title("Numerical Data")
    ax.set_xlabel("Index")
    ax.set_ylabel("Value")
    
    ax.grid()
    plt.show()



.. image:: output_43_0.png


.. code:: ipython3

    #S231
    
    import pandas as pd
    
    products = ["Pen", "Book", "Bag", "Bottle", "Pencil"]
    prices = [20, 50, 500, 100, 10]
    
    df = pd.DataFrame({
        "Product": products,
        "Price": prices
    })
    
    print("DataFrame:")
    print(df)
    
    print("Maximum Price:", df["Price"].max())
    print("Minimum Price:", df["Price"].min())
    print("Average Price:", df["Price"].mean())


.. parsed-literal::

    DataFrame:
      Product  Price
    0     Pen     20
    1    Book     50
    2     Bag    500
    3  Bottle    100
    4  Pencil     10
    Maximum Price: 500
    Minimum Price: 10
    Average Price: 136.0
    

.. code:: ipython3

    #S232
    
    import numpy as np
    
    data = np.array([
        [1, 2, 3],
        [4, 5, 6],
        [2, 3, 5]
    ])
    
    print("Original Data:")
    print(data)
    
    # L1 Normalization
    normalized = data / np.sum(np.abs(data), axis=1, keepdims=True)
    
    print("\nL1 Normalized Data:")
    print(normalized)


.. parsed-literal::

    Original Data:
    [[1 2 3]
     [4 5 6]
     [2 3 5]]
    
    L1 Normalized Data:
    [[0.16666667 0.33333333 0.5       ]
     [0.26666667 0.33333333 0.4       ]
     [0.2        0.3        0.5       ]]
    

.. code:: ipython3

    #S241
    
    import pandas as pd
    
    data = {
        "Name": ["Amit", "Rahul", "Sneha", "Priya", "Rohan"],
        "Study_Hours": [2, 4, 3, 5, 6],
        "Marks": [45, 70, 60, 85, 90]
    }
    
    df = pd.DataFrame(data)
    
    print("First 3 Rows:")
    print(df.head(3))
    
    print("\nLast 3 Rows:")
    print(df.tail(3))
    
    print("\nShape:")
    print(df.shape)
    


.. parsed-literal::

    First 3 Rows:
        Name  Study_Hours  Marks
    0   Amit            2     45
    1  Rahul            4     70
    2  Sneha            3     60
    
    Last 3 Rows:
        Name  Study_Hours  Marks
    2  Sneha            3     60
    3  Priya            5     85
    4  Rohan            6     90
    
    Shape:
    (5, 3)
    

.. code:: ipython3

    #S243
    
    import pandas as pd
    
    df = pd.read_csv("Student_Marks.csv")
    
    print("Before Discretization:")
    print(df)
    
    bins = [0, 40, 60, 80, 100]
    labels = ["Poor", "Average", "Good", "Excellent"]
    
    df["Performance"] = pd.cut(
        df["Marks"],
        bins=bins,
        labels=labels
    )
    
    print("\nAfter Discretization:")
    print(df)


::


    ---------------------------------------------------------------------------

    FileNotFoundError                         Traceback (most recent call last)

    Cell In[26], line 5
          1 #S243
          3 import pandas as pd
    ----> 5 df = pd.read_csv("Student_Marks.csv")
          7 print("Before Discretization:")
          8 print(df)
    

    File D:\Anaconda\Lib\site-packages\pandas\io\parsers\readers.py:1026, in read_csv(filepath_or_buffer, sep, delimiter, header, names, index_col, usecols, dtype, engine, converters, true_values, false_values, skipinitialspace, skiprows, skipfooter, nrows, na_values, keep_default_na, na_filter, verbose, skip_blank_lines, parse_dates, infer_datetime_format, keep_date_col, date_parser, date_format, dayfirst, cache_dates, iterator, chunksize, compression, thousands, decimal, lineterminator, quotechar, quoting, doublequote, escapechar, comment, encoding, encoding_errors, dialect, on_bad_lines, delim_whitespace, low_memory, memory_map, float_precision, storage_options, dtype_backend)
       1013 kwds_defaults = _refine_defaults_read(
       1014     dialect,
       1015     delimiter,
       (...)   1022     dtype_backend=dtype_backend,
       1023 )
       1024 kwds.update(kwds_defaults)
    -> 1026 return _read(filepath_or_buffer, kwds)
    

    File D:\Anaconda\Lib\site-packages\pandas\io\parsers\readers.py:620, in _read(filepath_or_buffer, kwds)
        617 _validate_names(kwds.get("names", None))
        619 # Create the parser.
    --> 620 parser = TextFileReader(filepath_or_buffer, **kwds)
        622 if chunksize or iterator:
        623     return parser
    

    File D:\Anaconda\Lib\site-packages\pandas\io\parsers\readers.py:1620, in TextFileReader.__init__(self, f, engine, **kwds)
       1617     self.options["has_index_names"] = kwds["has_index_names"]
       1619 self.handles: IOHandles | None = None
    -> 1620 self._engine = self._make_engine(f, self.engine)
    

    File D:\Anaconda\Lib\site-packages\pandas\io\parsers\readers.py:1880, in TextFileReader._make_engine(self, f, engine)
       1878     if "b" not in mode:
       1879         mode += "b"
    -> 1880 self.handles = get_handle(
       1881     f,
       1882     mode,
       1883     encoding=self.options.get("encoding", None),
       1884     compression=self.options.get("compression", None),
       1885     memory_map=self.options.get("memory_map", False),
       1886     is_text=is_text,
       1887     errors=self.options.get("encoding_errors", "strict"),
       1888     storage_options=self.options.get("storage_options", None),
       1889 )
       1890 assert self.handles is not None
       1891 f = self.handles.handle
    

    File D:\Anaconda\Lib\site-packages\pandas\io\common.py:873, in get_handle(path_or_buf, mode, encoding, compression, memory_map, is_text, errors, storage_options)
        868 elif isinstance(handle, str):
        869     # Check whether the filename is to be opened in binary mode.
        870     # Binary mode does not support 'encoding' and 'newline'.
        871     if ioargs.encoding and "b" not in ioargs.mode:
        872         # Encoding
    --> 873         handle = open(
        874             handle,
        875             ioargs.mode,
        876             encoding=ioargs.encoding,
        877             errors=errors,
        878             newline="",
        879         )
        880     else:
        881         # Binary mode
        882         handle = open(handle, ioargs.mode)
    

    FileNotFoundError: [Errno 2] No such file or directory: 'Student_Marks.csv'


.. code:: ipython3

    #S251
    
    import pandas as pd
    
    data = {
        "Product": ["Pen", "Book", "Bag", "Bottle", "Pencil"],
        "Quantity": [10, 5, 3, 8, 15],
        "Price": [20, 50, 500, 100, 10]
    }
    
    df = pd.DataFrame(data)
    
    print("DataFrame:")
    print(df)
    
    print("\nTotal Quantity:", df["Quantity"].sum())
    print("Average Price:", df["Price"].mean())


.. parsed-literal::

    DataFrame:
      Product  Quantity  Price
    0     Pen        10     20
    1    Book         5     50
    2     Bag         3    500
    3  Bottle         8    100
    4  Pencil        15     10
    
    Total Quantity: 41
    Average Price: 136.0
    

.. code:: ipython3

    #S253
    
    import numpy as np
    
    data = np.array([
        [3, 4],
        [5, 12],
        [8, 6]
    ])
    
    print("Original Data:")
    print(data)
    
    # Unit-length normalization
    norm = np.sqrt(np.sum(data ** 2, axis=1, keepdims=True))
    normalized = data / norm
    
    print("\nNormalized Data:")
    print(normalized)


.. parsed-literal::

    Original Data:
    [[ 3  4]
     [ 5 12]
     [ 8  6]]
    
    Normalized Data:
    [[0.6        0.8       ]
     [0.38461538 0.92307692]
     [0.8        0.6       ]]
    

.. code:: ipython3

    #S261
    
    import numpy as np
    from sklearn import preprocessing
    
    data = np.array([10, 20, 30, 40, 50])
    
    print("Original Data:")
    print(data)
    
    standardized = preprocessing.scale(data)
    
    print("\nStandardized Data:")
    print(standardized)


.. parsed-literal::

    Original Data:
    [10 20 30 40 50]
    
    Standardized Data:
    [-1.41421356 -0.70710678  0.          0.70710678  1.41421356]
    

.. code:: ipython3

    #S263
    
    import pandas as pd
    import seaborn as sns
    import matplotlib.pyplot as plt
    
    df = pd.read_csv("iris.csv")
    
    sns.pairplot(df, hue="Species")
    
    plt.show()

.. code:: ipython3

    #S271
    
    import pandas as pd
    
    data = {
        "Category": ["A", "B", "C", "D", "E"],
        "Value": [20, 50, 30, 80, 40]
    }
    
    df = pd.DataFrame(data)
    
    print("DataFrame:")
    print(df)
    
    print("\nMaximum Value:", df["Value"].max())
    print("Minimum Value:", df["Value"].min())


.. parsed-literal::

    DataFrame:
      Category  Value
    0        A     20
    1        B     50
    2        C     30
    3        D     80
    4        E     40
    
    Maximum Value: 80
    Minimum Value: 20
    

.. code:: ipython3

    #S273
    
    import pandas as pd
    import seaborn as sns
    import matplotlib.pyplot as plt
    
    df = pd.read_csv("iris.csv")
    
    features = [
        "SepalLengthCm",
        "SepalWidthCm",
        "PetalLengthCm",
        "PetalWidthCm"
    ]
    
    for feature in features:
        sns.boxplot(data=df, x="Species", y=feature)
        plt.title(feature)
        plt.show()
